#--coding:utf-8--

from lxml import etree

text = """
<div>
    <ul>
         <li class="item-0"><a href="link1.html">first item</a></li>
         <li class="item-1"><a href="link2.html">second item</a></li>
         <li class="item-inactive"><a href="link3.html"><span class="bold">third item</span></a></li>
         <li class="item-1"><a href="link4.html">fourth item</a></li>
         <li class="item-0"><a href="link5.html">fifth item</a></li>
    </ul>
 </div>
"""


# 将字符串解析为html文档
# html = etree.HTML(text)
# print(html)
# 按字符串序列号化html
# result = etree.tostring(html).decode('utf-8')
# print(result)

# 读取
html = etree.parse('hello.html')
result = etree.tostring(html).decode('utf-8')
print(result)
